# -*- coding: utf-8 -*-
"""
Intangible Cultural Heritage Traditional Sports Project - icourse163 Course Search Crawler (Improved Version)
"""

import os
import re
import time
import pandas as pd
from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from selenium.common.exceptions import TimeoutException, NoSuchElementException, WebDriverException
from selenium.webdriver.chrome.service import Service
from selenium.webdriver.chrome.options import Options

# ==================== Configuration ====================
INPUT_FILE = "非遗传统体育项目清单.xlsx"
OUTPUT_FILE = "检索结果1_国家智慧教育平台非遗体育课程.xlsx"
SEARCH_URL = "https://www.icourse163.org/search.htm"
REQUEST_DELAY = 3
PAGE_LOAD_TIMEOUT = 20

# ==================== Helper Functions ====================

def clean_project_name(name):
    """Remove parentheses and their inner content from project name"""
    if not isinstance(name, str):
        return str(name)
    cleaned = re.sub(r'（.*?）', '', name)
    cleaned = re.sub(r'\(.*?\)', '', cleaned)
    return cleaned.strip()


def setup_driver():
    """Launch ChromeDriver, load chromedriver.exe from script directory with highest priority"""
    chrome_options = Options()
    # chrome_options.add_argument("--headless")  # Uncomment to enable headless invisible‑browser mode
    chrome_options.add_argument("--disable-gpu")
    chrome_options.add_argument("--no-sandbox")
    chrome_options.add_argument("--disable-dev-shm-usage")
    chrome_options.add_argument(
        "user-agent=Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
        "AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
    )
    chrome_options.add_experimental_option("excludeSwitches", ["enable-automation"])
    chrome_options.add_experimental_option("useAutomationExtension", False)

    chromedriver_path = os.path.join(os.path.dirname(__file__), "chromedriver.exe")
    if os.path.exists(chromedriver_path):
        print(f"✅ Using ChromeDriver: {chromedriver_path}")
        service = Service(chromedriver_path)
    else:
        print("⚠️ chromedriver.exe not found, attempting to load from system PATH")
        service = Service()

    driver = webdriver.Chrome(service=service, options=chrome_options)
    driver.execute_cdp_cmd("Page.addScriptToEvaluateOnNewDocument", {
        "source": """
            Object.defineProperty(navigator, 'webdriver', {
                get: () => undefined
            })
        """
    })
    driver.set_page_load_timeout(PAGE_LOAD_TIMEOUT)
    return driver


def search_courses(driver, keyword):
    """
    Search target keyword and parse course information, return a list of matched course dictionaries
    """
    url = f"{SEARCH_URL}?search={keyword}"
    print(f"  Visiting: {url}")

    try:
        driver.get(url)
        # Wait until webpage finishes loading
        WebDriverWait(driver, 10).until(
            lambda d: d.execute_script("return document.readyState") == "complete"
        )
        time.sleep(2)  # Extra waiting time for dynamic‑rendered content

        # Detect empty‑search‑result prompt
        if "没有找到相关课程" in driver.page_source or "暂无课程" in driver.page_source:
            print(f"  ⚠️ No relevant courses found")
            return []

    except TimeoutException:
        print(f"  ⚠️ Page loading timeout")
        return []
    except WebDriverException as e:
        print(f"  ⚠️ WebDriver Exception: {type(e).__name__} - {e}")
        return []
    except Exception as e:
        print(f"  ⚠️ Access Exception: {type(e).__name__} - {e}")
        return []

    courses = []
    try:
        # Wait for at least one course‑card element, use multi‑selector fallback strategy
        wait = WebDriverWait(driver, 10)
        # Try multiple candidate CSS selectors
        course_elements = []
        selectors = [
            ".course-card",
            ".u-course",
            ".m-course",
            "[class*='course-item']",
            ".g-mn .u-course",
            ".j-list .u-course"
        ]
        for selector in selectors:
            try:
                wait.until(EC.presence_of_element_located((By.CSS_SELECTOR, selector)))
                elements = driver.find_elements(By.CSS_SELECTOR, selector)
                if elements:
                    course_elements = elements
                    print(f"  Found {len(elements)} courses (selector: {selector})")
                    break
            except TimeoutException:
                continue

        if not course_elements:
            # Fallback: locate elements by XPath containing course feature keywords
            try:
                course_elements = driver.find_elements(By.XPATH,
                    "//div[contains(@class, 'course') and .//span[contains(text(), '人参加')]]")
                if course_elements:
                    print(f"  Found {len(course_elements)} courses via XPath")
            except:
                pass

        if not course_elements:
            print(f"  ⚠️ No course elements captured, page‑structure changes or loading failure may occur")
            # Optional: print page title and partial source code for debugging
            print(f"     Page title: {driver.title}")
            return []

        # Parse each course card, extract at most the first 10 records
        for elem in course_elements[:10]:
            try:
                info = {"检索关键词": keyword}

                # ---- Course Name ----
                name = "未知"
                for tag in ['h3', 'h4', '.course-name', '.u-course-name', 'a[title]']:
                    try:
                        if tag.startswith('a'):
                            name_elem = elem.find_element(By.CSS_SELECTOR, tag)
                            name = name_elem.get_attribute("title").strip()
                        else:
                            name_elem = elem.find_element(By.CSS_SELECTOR, tag)
                            name = name_elem.text.strip()
                        if name:
                            break
                    except:
                        continue
                info["课程名称"] = name

                # ---- Offering University ----
                school = "未知"
                for tag in ['.school', '.u-school', '.course-school', '.t21', '[class*="school"]']:
                    try:
                        school_elem = elem.find_element(By.CSS_SELECTOR, tag)
                        school = school_elem.text.strip()
                        if school:
                            break
                    except:
                        continue
                info["开课高校"] = school

                # ---- Course Instructor ----
                teacher = "未知"
                for tag in ['.teacher', '.u-teacher', '.course-teacher', '.t22', '[class*="teacher"]']:
                    try:
                        teacher_elem = elem.find_element(By.CSS_SELECTOR, tag)
                        teacher = teacher_elem.text.strip()
                        if teacher:
                            break
                    except:
                        continue
                info["课程负责人"] = teacher

                # ---- Course Type (online by default) ----
                info["课程类型"] = "线上"
                try:
                    labels = elem.find_elements(By.CSS_SELECTOR, ".tag, .u-tag, .m-tag, [class*='tag']")
                    for label in labels:
                        txt = label.text.strip()
                        if "混合" in txt:
                            info["课程类型"] = "线上线下混合"
                            break
                        elif "线下" in txt:
                            info["课程类型"] = "线下"
                            break
                except:
                    pass

                # ---- Course Level ----
                level = "校级"
                try:
                    labels = elem.find_elements(By.CSS_SELECTOR, ".tag, .u-tag, .m-tag, [class*='tag']")
                    for label in labels:
                        txt = label.text.strip()
                        if "国家级" in txt or "一流" in txt:
                            level = "国家级一流本科课程"
                            break
                        elif "省级" in txt:
                            level = "省级一流"
                            break
                except:
                    pass
                info["课程等级"] = level

                # ---- Course Introduction ----
                desc = "暂无简介"
                for tag in ['.desc', '.u-desc', '.course-desc', '.m-desc', '[class*="desc"]']:
                    try:
                        desc_elem = elem.find_element(By.CSS_SELECTOR, tag)
                        desc = desc_elem.text.strip()
                        if desc:
                            break
                    except:
                        continue
                if desc == "暂无简介" or not desc:
                    desc = elem.text.strip()[:100]
                info["课程简介"] = desc

                courses.append(info)

            except Exception as e:
                print(f"  ⚠️ Error parsing single‑course element: {type(e).__name__} - {e}")
                continue

    except Exception as e:
        print(f"  ⚠️ Failed to fetch course‑list data: {type(e).__name__} - {e}")
        return []

    return courses


def main():
    print("=" * 70)
    print("Intangible Cultural Heritage Traditional Sports Project - icourse163 Course Search Crawler (Improved Version)")
    print("=" * 70)

    # Check whether input file exists
    if not os.path.exists(INPUT_FILE):
        print(f"❌ File '{INPUT_FILE}' does not exist")
        return

    try:
        df = pd.read_excel(INPUT_FILE)
        print(f"✅ File loaded successfully, total items: {len(df)}")
    except Exception as e:
        print(f"❌ Excel reading failed: {e}")
        return

    if '类型' not in df.columns or '名称' not in df.columns:
        print("❌ Required columns '类型' or '名称' are missing")
        return

    # Filter newly‑added projects
    df_new = df[df['类型'] == '新增项目'].copy()
    print(f"✅ Count of newly‑added projects: {len(df_new)}")

    if len(df_new) == 0:
        print("⚠️ No newly‑added projects found")
        return

    # Clean project names and remove duplicate keywords
    df_new['名称_清洗'] = df_new['名称'].apply(clean_project_name)
    df_new = df_new.drop_duplicates(subset=['名称_清洗'])
    project_names = df_new['名称_清洗'].tolist()
    print(f"✅ Distinct search‑keyword quantity after deduplication: {len(project_names)}")
    print(f"Keyword sample: {project_names[:5]}")

    # Launch Chrome browser instance
    print("\n🚀 Starting Chrome...")
    try:
        driver = setup_driver()
        print("✅ Browser launched successfully")
    except Exception as e:
        print(f"❌ Browser startup failed: {e}")
        return

    # Iterate and execute search for every keyword
    all_results = []
    total = len(project_names)
    print("\n" + "=" * 70)
    print("Start searching...")
    print("=" * 70)

    for idx, keyword in enumerate(project_names, 1):
        print(f"\n[{idx}/{total}] Search keyword: {keyword}")
        courses = search_courses(driver, keyword)
        if courses:
            print(f"  ✅ {len(courses)} matched courses found")
            all_results.extend(courses)
        else:
            print(f"  ⚠️ No courses retrieved")
        if idx < total:
            print(f"  ⏳ Sleep {REQUEST_DELAY} seconds before next request")
            time.sleep(REQUEST_DELAY)

    driver.quit()
    print("\n✅ Chrome browser closed")

    # Save crawled dataset into excel
    if all_results:
        df_res = pd.DataFrame(all_results)
        df_res = df_res.drop_duplicates()
        cols = ["检索关键词", "课程名称", "开课高校", "课程负责人", "课程类型", "课程等级", "课程简介"]
        df_res = df_res[cols]
        df_res.to_excel(OUTPUT_FILE, index=False, engine='openpyxl')
        print(f"\n✅ Search‑result saved: {os.path.abspath(OUTPUT_FILE)}")
        print(f"   Total valid records: {len(df_res)}")
    else:
        print("\n⚠️ Zero valid courses retrieved")

    print("\n" + "=" * 70)
    print(f"Task completed! Searched‑project count: {total}, captured‑course records: {len(all_results)}")
    print("=" * 70)


if __name__ == "__main__":
    main()
